agora inbox for pgsql-hackers@postgresql.org
help / color / mirror / Atom feed[PATCH v2 5/8] jit: explain: remove backend lifetime module count from function name.
590+ messages / 2 participants
[nested] [flat]
* [PATCH v2 5/8] jit: explain: remove backend lifetime module count from function name.
@ 2019-09-26 21:05 Andres Freund <andres@anarazel.de>
0 siblings, 0 replies; 590+ messages in thread
From: Andres Freund @ 2019-09-26 21:05 UTC (permalink / raw)
Also expand function name to include in which module the function is -
without that it's harder to analyze which functions were emitted
separately (a performance concern).
Author:
Reviewed-By:
Discussion: https://postgr.es/m/
Backpatch:
---
src/backend/commands/explain.c | 65 +++++++++++++++++++++++++++++-----
src/backend/jit/llvm/llvmjit.c | 18 +++++++---
src/include/jit/llvmjit.h | 5 ++-
3 files changed, 75 insertions(+), 13 deletions(-)
diff --git a/src/backend/commands/explain.c b/src/backend/commands/explain.c
index 3ccb76bdfd1..02455865d9f 100644
--- a/src/backend/commands/explain.c
+++ b/src/backend/commands/explain.c
@@ -2228,6 +2228,43 @@ show_expression(Node *node, ExprState *expr, const char *qlabel,
}
}
+/*
+ * To make JIT explain output reproducible, remove the module generation from
+ * function names. That makes it a bit harder to correlate with profiles etc,
+ * but reproducability is more important.
+ */
+static char *
+jit_funcname_for_display(const char *funcname)
+{
+ int func_counter; /* nth function in query */
+ size_t mod_num; /* nth module in query */
+ size_t mod_generation; /* nth module in backend */
+ int basename_end;
+ int matchcount = 0;
+
+ /*
+ * The pattern we need to match, see llvm_expand_funcname, is
+ * "%s_%zu_%d_mod_%zu". Find the fourth _ from the end, so a _ in the name
+ * is OK.
+ */
+ for (basename_end = strlen(funcname); basename_end >= 0; basename_end--)
+ {
+ if (funcname[basename_end] == '_' && ++matchcount == 4)
+ break;
+ }
+
+ /* couldn't parse, bail out */
+ if (matchcount != 4)
+ return pstrdup(funcname);
+
+ /* couldn't parse, bail out */
+ if (sscanf(funcname + basename_end, "_%zu_%d_mod_%zu",
+ &mod_num, &func_counter, &mod_generation) != 3)
+ return pstrdup(funcname);
+
+ return psprintf("%s_%zu_%d", pnstrdup(funcname, basename_end), mod_num, func_counter);
+}
+
static void
show_jit_expr_details(ExprState *expr, ExplainState *es)
{
@@ -2239,7 +2276,8 @@ show_jit_expr_details(ExprState *expr, ExplainState *es)
if (es->format == EXPLAIN_FORMAT_TEXT)
{
if (expr->flags & EEO_FLAG_JIT_EXPR)
- appendStringInfo(es->str, "JIT-Expr: %s", expr->expr_funcname);
+ appendStringInfo(es->str, "JIT-Expr: %s",
+ jit_funcname_for_display(expr->expr_funcname));
else
appendStringInfoString(es->str, "JIT-Expr: false");
@@ -2250,19 +2288,22 @@ show_jit_expr_details(ExprState *expr, ExplainState *es)
*/
if (expr->scan_funcname)
- appendStringInfo(es->str, ", JIT-Deform-Scan: %s", expr->scan_funcname);
+ appendStringInfo(es->str, ", JIT-Deform-Scan: %s",
+ jit_funcname_for_display(expr->scan_funcname));
else if (expr->flags & EEO_FLAG_JIT_EXPR &&
expr->flags & EEO_FLAG_DEFORM_SCAN)
appendStringInfo(es->str, ", JIT-Deform-Scan: false");
if (expr->outer_funcname)
- appendStringInfo(es->str, ", JIT-Deform-Outer: %s", expr->outer_funcname);
+ appendStringInfo(es->str, ", JIT-Deform-Outer: %s",
+ jit_funcname_for_display(expr->outer_funcname));
else if (expr->flags & EEO_FLAG_JIT_EXPR &&
expr->flags & EEO_FLAG_DEFORM_OUTER)
appendStringInfo(es->str, ", JIT-Deform-Outer: false");
if (expr->inner_funcname)
- appendStringInfo(es->str, ", JIT-Deform-Inner: %s", expr->inner_funcname);
+ appendStringInfo(es->str, ", JIT-Deform-Inner: %s",
+ jit_funcname_for_display(expr->inner_funcname));
else if (expr->flags & EEO_FLAG_JIT_EXPR &&
expr->flags & (EEO_FLAG_DEFORM_INNER))
appendStringInfo(es->str, ", JIT-Deform-Inner: false");
@@ -2270,26 +2311,34 @@ show_jit_expr_details(ExprState *expr, ExplainState *es)
else
{
if (expr->flags & EEO_FLAG_JIT_EXPR)
- ExplainPropertyText("JIT-Expr", expr->expr_funcname, es);
+ ExplainPropertyText("JIT-Expr",
+ jit_funcname_for_display(expr->expr_funcname),
+ es);
else
ExplainPropertyBool("JIT-Expr", false, es);
if (expr->scan_funcname)
- ExplainProperty("JIT-Deform-Scan", NULL, expr->scan_funcname, false, es);
+ ExplainProperty("JIT-Deform-Scan", NULL,
+ jit_funcname_for_display(expr->scan_funcname),
+ false, es);
else if (expr->flags & EEO_FLAG_DEFORM_SCAN)
ExplainProperty("JIT-Deform-Scan", NULL, "false", true, es);
else
ExplainProperty("JIT-Deform-Scan", NULL, "null", true, es);
if (expr->outer_funcname)
- ExplainProperty("JIT-Deform-Outer", NULL, expr->outer_funcname, false, es);
+ ExplainProperty("JIT-Deform-Outer", NULL,
+ jit_funcname_for_display(expr->outer_funcname),
+ false, es);
else if (expr->flags & EEO_FLAG_DEFORM_OUTER)
ExplainProperty("JIT-Deform-Outer", NULL, "false", true, es);
else
ExplainProperty("JIT-Deform-Outer", NULL, "null", true, es);
if (expr->inner_funcname)
- ExplainProperty("JIT-Deform-Inner", NULL, expr->inner_funcname, false, es);
+ ExplainProperty("JIT-Deform-Inner", NULL,
+ jit_funcname_for_display(expr->inner_funcname),
+ false, es);
else if (expr->flags & EEO_FLAG_DEFORM_INNER)
ExplainProperty("JIT-Deform-Inner", NULL, "false", true, es);
else
diff --git a/src/backend/jit/llvm/llvmjit.c b/src/backend/jit/llvm/llvmjit.c
index 5489e118041..177a00f3826 100644
--- a/src/backend/jit/llvm/llvmjit.c
+++ b/src/backend/jit/llvm/llvmjit.c
@@ -227,6 +227,8 @@ llvm_mutable_module(LLVMJitContext *context)
char *
llvm_expand_funcname(struct LLVMJitContext *context, const char *basename)
{
+ char *funcname;
+
Assert(context->module != NULL);
context->base.instr.created_functions++;
@@ -234,11 +236,19 @@ llvm_expand_funcname(struct LLVMJitContext *context, const char *basename)
/*
* Previously we used dots to separate, but turns out some tools, e.g.
* GDB, don't like that and truncate name.
+ *
+ * Append the backend-lifetime module count to the end, so it's easier for
+ * humans and machines to compare the generated function names across
+ * queries, the prefix will be the same from query execution to query
+ * execution.
*/
- return psprintf("%s_%zu_%d",
- basename,
- context->module_generation,
- context->counter++);
+ funcname = psprintf("%s_%zu_%d_mod_%zu",
+ basename,
+ context->base.instr.created_modules - 1,
+ context->counter++,
+ context->module_generation);
+
+ return funcname;
}
/*
diff --git a/src/include/jit/llvmjit.h b/src/include/jit/llvmjit.h
index 6178864b2e6..e45ff99194f 100644
--- a/src/include/jit/llvmjit.h
+++ b/src/include/jit/llvmjit.h
@@ -41,7 +41,10 @@ typedef struct LLVMJitContext
{
JitContext base;
- /* number of modules created */
+ /*
+ * llvm_generation when ->module was created, monotonically increasing
+ * within the lifetime of a backend.
+ */
size_t module_generation;
/* current, "open for write", module */
--
2.23.0.385.gbc12974a89
--ga6shgqrocqphdjc
Content-Type: text/x-diff; charset=us-ascii
Content-Disposition: attachment;
filename="v2-0006-WIP-explain-Show-per-phase-information-about-aggr.patch"
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Xf36GNcW+0xZNily
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v2-0009-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
* [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics
@ 2026-07-09 20:21 Nathan Bossart <nathan@postgresql.org>
0 siblings, 0 replies; 590+ messages in thread
From: Nathan Bossart @ 2026-07-09 20:21 UTC (permalink / raw)
---
src/backend/access/heap/heapam_handler.c | 2 +-
src/backend/access/table/tableam.c | 59 +++++++++++-------------
src/include/access/relscan.h | 8 ++--
3 files changed, 31 insertions(+), 38 deletions(-)
diff --git a/src/backend/access/heap/heapam_handler.c b/src/backend/access/heap/heapam_handler.c
index bf87430cf01..0f24a132564 100644
--- a/src/backend/access/heap/heapam_handler.c
+++ b/src/backend/access/heap/heapam_handler.c
@@ -1965,7 +1965,7 @@ heapam_scan_get_blocks_done(HeapScanDesc hscan)
if (hscan->rs_base.rs_parallel != NULL)
{
bpscan = (ParallelBlockTableScanDesc) hscan->rs_base.rs_parallel;
- startblock = bpscan->phs_startblock;
+ startblock = pg_atomic_read_u32(&bpscan->phs_startblock);
}
else
startblock = hscan->rs_startblock;
diff --git a/src/backend/access/table/tableam.c b/src/backend/access/table/tableam.c
index 68ff0966f1c..f2038ea9205 100644
--- a/src/backend/access/table/tableam.c
+++ b/src/backend/access/table/tableam.c
@@ -421,9 +421,8 @@ table_block_parallelscan_initialize(Relation rel, ParallelTableScanDesc pscan)
bpscan->base.phs_syncscan = synchronize_seqscans &&
!RelationUsesLocalBuffers(rel) &&
bpscan->phs_nblocks > NBuffers / 4;
- SpinLockInit(&bpscan->phs_mutex);
- bpscan->phs_startblock = InvalidBlockNumber;
- bpscan->phs_numblock = InvalidBlockNumber;
+ pg_atomic_init_u32(&bpscan->phs_startblock, InvalidBlockNumber);
+ pg_atomic_init_u32(&bpscan->phs_numblock, InvalidBlockNumber);
pg_atomic_init_u64(&bpscan->phs_nallocated, 0);
return sizeof(ParallelBlockTableScanDescData);
@@ -459,25 +458,22 @@ table_block_parallelscan_startblock_init(Relation rel,
StaticAssertDecl(MaxBlockNumber <= 0xFFFFFFFE,
"pg_nextpower2_32 may be too small for non-standard BlockNumber width");
- BlockNumber sync_startpage = InvalidBlockNumber;
BlockNumber scan_nblocks;
/* Reset the state we use for controlling allocation size. */
memset(pbscanwork, 0, sizeof(*pbscanwork));
-retry:
- /* Grab the spinlock. */
- SpinLockAcquire(&pbscan->phs_mutex);
-
/*
* When the caller specified a limit on the number of blocks to scan, set
* that in the ParallelBlockTableScanDesc, if it's not been done by
* another worker already.
*/
- if (numblocks != InvalidBlockNumber &&
- pbscan->phs_numblock == InvalidBlockNumber)
+ if (numblocks != InvalidBlockNumber)
{
- pbscan->phs_numblock = numblocks;
+ uint32 expected = InvalidBlockNumber;
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_numblock, &expected,
+ numblocks);
}
/*
@@ -485,36 +481,35 @@ retry:
* so now. If a startblock was specified, start there, otherwise if this
* is not a synchronized scan, we just start at block 0, but if it is a
* synchronized scan, we must get the starting position from the
- * synchronized scan machinery. We can't hold the spinlock while doing
- * that, though, so release the spinlock, get the information we need, and
- * retry. If nobody else has initialized the scan in the meantime, we'll
- * fill in the value we fetched on the second time through.
+ * synchronized scan machinery.
+ *
+ * If another worker initializes phs_startblock concurrently, just use
+ * their value.
*/
- if (pbscan->phs_startblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_startblock) == InvalidBlockNumber)
{
+ BlockNumber newstartblock;
+ uint32 expected = InvalidBlockNumber;
+
if (startblock != InvalidBlockNumber)
- pbscan->phs_startblock = startblock;
+ newstartblock = startblock;
else if (!pbscan->base.phs_syncscan)
- pbscan->phs_startblock = 0;
- else if (sync_startpage != InvalidBlockNumber)
- pbscan->phs_startblock = sync_startpage;
+ newstartblock = 0;
else
- {
- SpinLockRelease(&pbscan->phs_mutex);
- sync_startpage = ss_get_location(rel, pbscan->phs_nblocks);
- goto retry;
- }
+ newstartblock = ss_get_location(rel, pbscan->phs_nblocks);
+
+ pg_atomic_compare_exchange_u32(&pbscan->phs_startblock, &expected,
+ newstartblock);
}
- SpinLockRelease(&pbscan->phs_mutex);
/*
* Figure out how many blocks we're going to scan; either all of them, or
* just phs_numblock's worth, if a limit has been imposed.
*/
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* We determine the chunk size based on scan_nblocks. First we split
@@ -595,10 +590,10 @@ table_block_parallelscan_nextpage(Relation rel,
*/
/* First, figure out how many blocks we're planning on scanning */
- if (pbscan->phs_numblock == InvalidBlockNumber)
+ if (pg_atomic_read_u32(&pbscan->phs_numblock) == InvalidBlockNumber)
scan_nblocks = pbscan->phs_nblocks;
else
- scan_nblocks = pbscan->phs_numblock;
+ scan_nblocks = pg_atomic_read_u32(&pbscan->phs_numblock);
/*
* Now check if we have any remaining blocks in a previous chunk for this
@@ -644,7 +639,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (nallocated >= scan_nblocks)
page = InvalidBlockNumber; /* all blocks have been allocated */
else
- page = (nallocated + pbscan->phs_startblock) % pbscan->phs_nblocks;
+ page = (nallocated + pg_atomic_read_u32(&pbscan->phs_startblock)) % pbscan->phs_nblocks;
/*
* Report scan location. Normally, we report the current page number.
@@ -658,7 +653,7 @@ table_block_parallelscan_nextpage(Relation rel,
if (page != InvalidBlockNumber)
ss_report_location(rel, page);
else if (nallocated == pbscan->phs_nblocks)
- ss_report_location(rel, pbscan->phs_startblock);
+ ss_report_location(rel, pg_atomic_read_u32(&pbscan->phs_startblock));
}
return page;
diff --git a/src/include/access/relscan.h b/src/include/access/relscan.h
index 2ea06a67a63..2305d0159f3 100644
--- a/src/include/access/relscan.h
+++ b/src/include/access/relscan.h
@@ -19,7 +19,6 @@
#include "nodes/tidbitmap.h"
#include "port/atomics.h"
#include "storage/relfilelocator.h"
-#include "storage/spin.h"
#include "utils/relcache.h"
@@ -99,10 +98,9 @@ typedef struct ParallelBlockTableScanDescData
ParallelTableScanDescData base;
BlockNumber phs_nblocks; /* # blocks in relation at start of scan */
- slock_t phs_mutex; /* mutual exclusion for setting startblock */
- BlockNumber phs_startblock; /* starting block number */
- BlockNumber phs_numblock; /* # blocks to scan, or InvalidBlockNumber if
- * no limit */
+ pg_atomic_uint32 phs_startblock; /* starting block number */
+ pg_atomic_uint32 phs_numblock; /* # blocks to scan, or InvalidBlockNumber
+ * if no limit */
pg_atomic_uint64 phs_nallocated; /* number of blocks allocated to
* workers so far. */
} ParallelBlockTableScanDescData;
--
2.50.1 (Apple Git-155)
--Ot5x3ffkWEuxilWO
Content-Type: text/plain; charset=us-ascii
Content-Disposition: attachment;
filename=v1-0008-convert-FastPathStrongRelationLocks-to-atomics.patch
^ permalink raw reply [nested|flat] 590+ messages in thread
end of thread, other threads:[~2026-07-09 20:21 UTC | newest]
Thread overview: 590+ messages (download: mbox mbox.gz follow: Atom feed)
-- links below jump to the message on this page --
2019-09-26 21:05 [PATCH v2 5/8] jit: explain: remove backend lifetime module count from function name. Andres Freund <andres@anarazel.de>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v2 8/9] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
2026-07-09 20:21 [PATCH v1 7/8] convert ParallelBlockTableScanDescData->phs_{start,num}block to atomics Nathan Bossart <nathan@postgresql.org>
This inbox is served by agora; see mirroring instructions
for how to clone and mirror all data and code used for this inbox